<!DOCTYPE html>
<html class="client-nojs vector-feature-night-mode-disabled vector-feature-language-in-header-enabled vector-feature-language-in-main-page-header-disabled vector-feature-page-tools-pinned-disabled vector-feature-toc-pinned-clientpref-1 vector-feature-main-menu-pinned-disabled vector-feature-limited-width-clientpref-1 vector-feature-limited-width-content-enabled vector-feature-custom-font-size-clientpref-1 vector-feature-appearance-pinned-clientpref-1 vector-sticky-header-enabled" lang="en" dir="ltr"><head>
<meta charset="UTF-8">
<title>Unsupervised learning</title>
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="canonical" href="https://en.wikipedia.org/wiki/Unsupervised_learning"> <link href="./mw/ext.cite.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/ext.math.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.icons.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.search.codex.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/user.styles.css" rel="stylesheet" type="text/css">
<meta name="ResourceLoaderDynamicStyles" content="">
<link rel="stylesheet" type="text/css" href="./mw/site.styles.css">
<link rel="stylesheet" type="text/css" href="./mw/noscript.css">
<link rel="stylesheet" type="text/css" href="./footer.css">
<link rel="stylesheet" type="text/css" href="./vector-2022.css">
</head>
<body class="skin--responsive skin-vector skin-vector-search-vue mediawiki ltr sitedir-ltr mw-hide-empty-elt ns-0 ns-subject page-Unsupervised_learning rootpage-Unsupervised_learning skin-vector-2022 action-view">
<div class="mw-page-container">
<div class="mw-page-container-inner">
<div class="mw-content-container">
<main id="content" class="mw-body">
<header class="mw-body-header vector-page-titlebar">
<h1 id="firstHeading" class="firstHeading mw-first-heading">
<span id="openzim-page-title" class="mw-page-title-main"><span class="mw-page-title-main">Unsupervised learning</span></span>
</h1>
</header>
<a id="top"></a>
<div id="bodyContent" class="vector-body ve-init-mw-desktopArticleTarget-targetContainer" aria-labelledby="firstHeading" data-mw-ve-target-container="">
<div id="mw-content-text" class="mw-body-content mw-content-ltr" lang="en" dir="ltr"><div class="mw-content-ltr mw-parser-output" lang="en" dir="ltr">
<style data-mw-deduplicate="TemplateStyles:r1129693374">
/* start https://en.wikipedia.org/ */
.mw-parser-output .hlist dl,.mw-parser-output .hlist ol,.mw-parser-output .hlist ul{margin:0;padding:0}.mw-parser-output .hlist dd,.mw-parser-output .hlist dt,.mw-parser-output .hlist li{margin:0;display:inline}.mw-parser-output .hlist.inline,.mw-parser-output .hlist.inline dl,.mw-parser-output .hlist.inline ol,.mw-parser-output .hlist.inline ul,.mw-parser-output .hlist dl dl,.mw-parser-output .hlist dl ol,.mw-parser-output .hlist dl ul,.mw-parser-output .hlist ol dl,.mw-parser-output .hlist ol ol,.mw-parser-output .hlist ol ul,.mw-parser-output .hlist ul dl,.mw-parser-output .hlist ul ol,.mw-parser-output .hlist ul ul{display:inline}.mw-parser-output .hlist .mw-empty-li{display:none}.mw-parser-output .hlist dt::after{content:": "}.mw-parser-output .hlist dd::after,.mw-parser-output .hlist li::after{content:" · ";font-weight:bold}.mw-parser-output .hlist dd:last-child::after,.mw-parser-output .hlist dt:last-child::after,.mw-parser-output .hlist li:last-child::after{content:none}.mw-parser-output .hlist dd dd:first-child::before,.mw-parser-output .hlist dd dt:first-child::before,.mw-parser-output .hlist dd li:first-child::before,.mw-parser-output .hlist dt dd:first-child::before,.mw-parser-output .hlist dt dt:first-child::before,.mw-parser-output .hlist dt li:first-child::before,.mw-parser-output .hlist li dd:first-child::before,.mw-parser-output .hlist li dt:first-child::before,.mw-parser-output .hlist li li:first-child::before{content:" (";font-weight:normal}.mw-parser-output .hlist dd dd:last-child::after,.mw-parser-output .hlist dd dt:last-child::after,.mw-parser-output .hlist dd li:last-child::after,.mw-parser-output .hlist dt dd:last-child::after,.mw-parser-output .hlist dt dt:last-child::after,.mw-parser-output .hlist dt li:last-child::after,.mw-parser-output .hlist li dd:last-child::after,.mw-parser-output .hlist li dt:last-child::after,.mw-parser-output .hlist li li:last-child::after{content:")";font-weight:normal}.mw-parser-output .hlist ol{counter-reset:listitem}.mw-parser-output .hlist ol>li{counter-increment:listitem}.mw-parser-output .hlist ol>li::before{content:" "counter(listitem)"\a0 "}.mw-parser-output .hlist dd ol>li:first-child::before,.mw-parser-output .hlist dt ol>li:first-child::before,.mw-parser-output .hlist li ol>li:first-child::before{content:" ("counter(listitem)"\a0 "}
/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r1246091330">
/* start https://en.wikipedia.org/ */
.mw-parser-output .sidebar{width:22em;float:right;clear:right;margin:0.5em 0 1em 1em;background:var(--background-color-neutral-subtle,#f8f9fa);border:1px solid var(--border-color-base,#a2a9b1);padding:0.2em;text-align:center;line-height:1.4em;font-size:88%;border-collapse:collapse;display:table}body.skin-minerva .mw-parser-output .sidebar{display:table!important;float:right!important;margin:0.5em 0 1em 1em!important}.mw-parser-output .sidebar-subgroup{width:100%;margin:0;border-spacing:0}.mw-parser-output .sidebar-left{float:left;clear:left;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-none{float:none;clear:both;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-outer-title{padding:0 0.4em 0.2em;font-size:125%;line-height:1.2em;font-weight:bold}.mw-parser-output .sidebar-top-image{padding:0.4em}.mw-parser-output .sidebar-top-caption,.mw-parser-output .sidebar-pretitle-with-top-image,.mw-parser-output .sidebar-caption{padding:0.2em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-pretitle{padding:0.4em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-title,.mw-parser-output .sidebar-title-with-pretitle{padding:0.2em 0.8em;font-size:145%;line-height:1.2em}.mw-parser-output .sidebar-title-with-pretitle{padding:0.1em 0.4em}.mw-parser-output .sidebar-image{padding:0.2em 0.4em 0.4em}.mw-parser-output .sidebar-heading{padding:0.1em 0.4em}.mw-parser-output .sidebar-content{padding:0 0.5em 0.4em}.mw-parser-output .sidebar-content-with-subgroup{padding:0.1em 0.4em 0.2em}.mw-parser-output .sidebar-above,.mw-parser-output .sidebar-below{padding:0.3em 0.8em;font-weight:bold}.mw-parser-output .sidebar-collapse .sidebar-above,.mw-parser-output .sidebar-collapse .sidebar-below{border-top:1px solid #aaa;border-bottom:1px solid #aaa}.mw-parser-output .sidebar-navbar{text-align:right;font-size:115%;padding:0 0.4em 0.4em}.mw-parser-output .sidebar-list-title{padding:0 0.4em;text-align:left;font-weight:bold;line-height:1.6em;font-size:105%}.mw-parser-output .sidebar-list-title-c{padding:0 0.4em;text-align:center;margin:0 3.3em}@media(max-width:640px){body.mediawiki .mw-parser-output .sidebar{width:100%!important;clear:both;float:none!important;margin-left:0!important;margin-right:0!important}}body.skin--responsive .mw-parser-output .sidebar a>img{max-width:none!important}@media screen{html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media print{body.ns-0 .mw-parser-output .sidebar{display:none!important}}
/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r886047488">
/* start https://en.wikipedia.org/ */
.mw-parser-output .nobold{font-weight:normal}
/* end https://en.wikipedia.org/ */
</style><table class="sidebar sidebar-collapse nomobile nowraplinks"><tbody><tr><td class="sidebar-pretitle">Part of a series on</td></tr><tr><th class="sidebar-title-with-pretitle"><a href="Machine_learning" title="Machine learning">Machine learning</a><br>and <a href="Data_mining" title="Data mining">data mining</a></th></tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Paradigms</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Supervised_learning" title="Supervised learning">Supervised learning</a></li>
<li><a href="Semi-supervised_learning" class="mw-redirect" title="Semi-supervised learning">Semi-supervised learning</a></li>
<li><a href="Self-supervised_learning" title="Self-supervised learning">Self-supervised learning</a></li>
<li><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></li>
<li><a href="Meta-learning_(computer_science)" title="Meta-learning (computer science)">Meta-learning</a></li>
<li><a href="Online_machine_learning" title="Online machine learning">Online learning</a></li>
<li><a href="Batch_learning" class="mw-redirect" title="Batch learning">Batch learning</a></li>
<li><a href="Curriculum_learning" title="Curriculum learning">Curriculum learning</a></li>
<li><a href="Rule-based_machine_learning" title="Rule-based machine learning">Rule-based learning</a></li>
<li><a href="Neuro-symbolic_AI" title="Neuro-symbolic AI">Neuro-symbolic AI</a></li>
<li><a href="Neuromorphic_engineering" class="mw-redirect" title="Neuromorphic engineering">Neuromorphic engineering</a></li>
<li><a href="Quantum_machine_learning" title="Quantum machine learning">Quantum machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Problems</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Statistical_classification" title="Statistical classification">Classification</a></li>
<li><a href="Generative_model" title="Generative model">Generative modeling</a></li>
<li><a href="Regression_analysis" title="Regression analysis">Regression</a></li>
<li><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></li>
<li><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></li>
<li><a href="Density_estimation" title="Density estimation">Density estimation</a></li>
<li><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></li>
<li><a href="Data_cleaning" class="mw-redirect" title="Data cleaning">Data cleaning</a></li>
<li><a href="Automated_machine_learning" title="Automated machine learning">AutoML</a></li>
<li><a href="Association_rule_learning" title="Association rule learning">Association rules</a></li>
<li><a href="Semantic_analysis_(machine_learning)" title="Semantic analysis (machine learning)">Semantic analysis</a></li>
<li><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></li>
<li><a href="Feature_engineering" title="Feature engineering">Feature engineering</a></li>
<li><a href="Feature_learning" title="Feature learning">Feature learning</a></li>
<li><a href="Learning_to_rank" title="Learning to rank">Learning to rank</a></li>
<li><a href="Grammar_induction" title="Grammar induction">Grammar induction</a></li>
<li><a href="Ontology_learning" title="Ontology learning">Ontology learning</a></li>
<li><a href="Multimodal_learning" title="Multimodal learning">Multimodal learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><div style="display: inline-block; line-height: 1.2em; padding: .1em 0;"><a href="Supervised_learning" title="Supervised learning">Supervised learning</a><br><span class="nobold"><span style="font-size: 85%;">(<b><a href="Statistical_classification" title="Statistical classification">classification</a></b> • <b><a href="Regression_analysis" title="Regression analysis">regression</a></b>)</span></span> </div></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Apprenticeship_learning" title="Apprenticeship learning">Apprenticeship learning</a></li>
<li><a href="Decision_tree_learning" title="Decision tree learning">Decision trees</a></li>
<li><a href="Ensemble_learning" title="Ensemble learning">Ensembles</a>
<ul><li><a href="Bootstrap_aggregating" title="Bootstrap aggregating">Bagging</a></li>
<li><a href="Boosting_(machine_learning)" title="Boosting (machine learning)">Boosting</a></li>
<li><a href="Random_forest" title="Random forest">Random forest</a></li></ul></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Linear_regression" title="Linear regression">Linear regression</a></li>
<li><a href="Naive_Bayes_classifier" title="Naive Bayes classifier">Naive Bayes</a></li>
<li><a href="Artificial_neural_network" class="mw-redirect" title="Artificial neural network">Artificial neural networks</a></li>
<li><a href="Logistic_regression" title="Logistic regression">Logistic regression</a></li>
<li><a href="Perceptron" title="Perceptron">Perceptron</a></li>
<li><a href="Relevance_vector_machine" title="Relevance vector machine">Relevance vector machine (RVM)</a></li>
<li><a href="Support_vector_machine" title="Support vector machine">Support vector machine (SVM)</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="BIRCH" title="BIRCH">BIRCH</a></li>
<li><a href="CURE_algorithm" title="CURE algorithm">CURE</a></li>
<li><a href="Hierarchical_clustering" title="Hierarchical clustering">Hierarchical</a></li>
<li><a href="K-means_clustering" title="K-means clustering"><i>k</i>-means</a></li>
<li><a href="Fuzzy_clustering" title="Fuzzy clustering">Fuzzy</a></li>
<li><a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization (EM)</a></li>
<li><br><a href="DBSCAN" title="DBSCAN">DBSCAN</a></li>
<li><a href="OPTICS_algorithm" title="OPTICS algorithm">OPTICS</a></li>
<li><a href="Mean_shift" title="Mean shift">Mean shift</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Factor_analysis" title="Factor analysis">Factor analysis</a></li>
<li><a href="Canonical_correlation" title="Canonical correlation">CCA</a></li>
<li><a href="Independent_component_analysis" title="Independent component analysis">ICA</a></li>
<li><a href="Linear_discriminant_analysis" title="Linear discriminant analysis">LDA</a></li>
<li><a href="Non-negative_matrix_factorization" title="Non-negative matrix factorization">NMF</a></li>
<li><a href="Principal_component_analysis" title="Principal component analysis">PCA</a></li>
<li><a href="Proper_generalized_decomposition" title="Proper generalized decomposition">PGD</a></li>
<li><a href="T-distributed_stochastic_neighbor_embedding" title="T-distributed stochastic neighbor embedding">t-SNE</a></li>
<li><a href="Sparse_dictionary_learning" title="Sparse dictionary learning">SDL</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Graphical_model" title="Graphical model">Graphical models</a>
<ul><li><a href="Bayesian_network" title="Bayesian network">Bayes net</a></li>
<li><a href="Conditional_random_field" title="Conditional random field">Conditional random field</a></li>
<li><a href="Hidden_Markov_model" title="Hidden Markov model">Hidden Markov</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Random_sample_consensus" title="Random sample consensus">RANSAC</a></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Local_outlier_factor" title="Local outlier factor">Local outlier factor</a></li>
<li><a href="Isolation_forest" title="Isolation forest">Isolation forest</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Neural_network_(machine_learning)" title="Neural network (machine learning)">Neural networks</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Autoencoder" title="Autoencoder">Autoencoder</a></li>
<li><a href="Deep_learning" title="Deep learning">Deep learning</a></li>
<li><a href="Feedforward_neural_network" title="Feedforward neural network">Feedforward neural network</a></li>
<li><a href="Recurrent_neural_network" title="Recurrent neural network">Recurrent neural network</a>
<ul><li><a href="Long_short-term_memory" title="Long short-term memory">LSTM</a></li>
<li><a href="Gated_recurrent_unit" title="Gated recurrent unit">GRU</a></li>
<li><a href="Echo_state_network" title="Echo state network">ESN</a></li>
<li><a href="Reservoir_computing" title="Reservoir computing">reservoir computing</a></li></ul></li>
<li><a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann machine</a>
<ul><li><a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">Restricted</a></li></ul></li>
<li><a href="Generative_adversarial_network" title="Generative adversarial network">GAN</a></li>
<li><a href="Diffusion_model" title="Diffusion model">Diffusion model</a></li>
<li><a href="Self-organizing_map" title="Self-organizing map">SOM</a></li>
<li><a href="Convolutional_neural_network" title="Convolutional neural network">Convolutional neural network</a>
<ul><li><a href="U-Net" title="U-Net">U-Net</a></li>
<li><a href="LeNet" title="LeNet">LeNet</a></li>
<li><a href="AlexNet" title="AlexNet">AlexNet</a></li>
<li><a href="DeepDream" title="DeepDream">DeepDream</a></li></ul></li>
<li><a href="Neural_field" title="Neural field">Neural field</a>
<ul><li><a href="Neural_radiance_field" title="Neural radiance field">Neural radiance field</a></li>
<li><a href="Physics-informed_neural_networks" title="Physics-informed neural networks">Physics-informed neural networks</a></li></ul></li>
<li><a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">Transformer</a>
<ul><li><a href="Vision_transformer" title="Vision transformer">Vision</a></li></ul></li>
<li><a href="Mamba_(deep_learning_architecture)" title="Mamba (deep learning architecture)">Mamba</a></li>
<li><a href="Spiking_neural_network" title="Spiking neural network">Spiking neural network</a></li>
<li><a href="Memtransistor" title="Memtransistor">Memtransistor</a></li>
<li><a href="Electrochemical_RAM" title="Electrochemical RAM">Electrochemical RAM</a> (ECRAM)</li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Q-learning" title="Q-learning">Q-learning</a></li>
<li><a href="Policy_gradient_method" title="Policy gradient method">Policy gradient</a></li>
<li><a href="State%E2%80%93action%E2%80%93reward%E2%80%93state%E2%80%93action" title="State–action–reward–state–action">SARSA</a></li>
<li><a href="Temporal_difference_learning" title="Temporal difference learning">Temporal difference (TD)</a></li>
<li><a href="Multi-agent_reinforcement_learning" title="Multi-agent reinforcement learning">Multi-agent</a>
<ul><li><a href="Self-play_(reinforcement_learning_technique)" class="mw-redirect" title="Self-play (reinforcement learning technique)">Self-play</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Learning with humans</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Active_learning_(machine_learning)" title="Active learning (machine learning)">Active learning</a></li>
<li><a href="Crowdsourcing" title="Crowdsourcing">Crowdsourcing</a></li>
<li><a href="Human-in-the-loop" title="Human-in-the-loop">Human-in-the-loop</a></li>
<li><a href="Mechanistic_interpretability" title="Mechanistic interpretability">Mechanistic interpretability</a></li>
<li><a href="Reinforcement_learning_from_human_feedback" title="Reinforcement learning from human feedback">RLHF</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Model diagnostics</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Coefficient_of_determination" title="Coefficient of determination">Coefficient of determination</a></li>
<li><a href="Confusion_matrix" title="Confusion matrix">Confusion matrix</a></li>
<li><a href="Learning_curve_(machine_learning)" title="Learning curve (machine learning)">Learning curve</a></li>
<li><a href="Receiver_operating_characteristic" title="Receiver operating characteristic">ROC curve</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Mathematical foundations</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Kernel_machines" class="mw-redirect" title="Kernel machines">Kernel machines</a></li>
<li><a href="Bias%E2%80%93variance_tradeoff" title="Bias–variance tradeoff">Bias–variance tradeoff</a></li>
<li><a href="Computational_learning_theory" title="Computational learning theory">Computational learning theory</a></li>
<li><a href="Empirical_risk_minimization" title="Empirical risk minimization">Empirical risk minimization</a></li>
<li><a href="Occam_learning" title="Occam learning">Occam learning</a></li>
<li><a href="Probably_approximately_correct_learning" title="Probably approximately correct learning">PAC learning</a></li>
<li><a href="Statistical_learning_theory" title="Statistical learning theory">Statistical learning</a></li>
<li><a href="Vapnik%E2%80%93Chervonenkis_theory" title="Vapnik–Chervonenkis theory">VC theory</a></li>
<li><a href="Topological_deep_learning" title="Topological deep learning">Topological deep learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Journals and conferences</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="AAAI_Conference_on_Artificial_Intelligence" title="AAAI Conference on Artificial Intelligence">AAAI</a></li>
<li><a href="ECML_PKDD" title="ECML PKDD">ECML PKDD</a></li>
<li><a href="Conference_on_Neural_Information_Processing_Systems" title="Conference on Neural Information Processing Systems">NeurIPS</a></li>
<li><a href="International_Conference_on_Machine_Learning" title="International Conference on Machine Learning">ICML</a></li>
<li><a href="International_Conference_on_Learning_Representations" title="International Conference on Learning Representations">ICLR</a></li>
<li><a href="International_Joint_Conference_on_Artificial_Intelligence" title="International Joint Conference on Artificial Intelligence">IJCAI</a></li>
<li><a href="Machine_Learning_(journal)" title="Machine Learning (journal)">ML</a></li>
<li><a href="Journal_of_Machine_Learning_Research" title="Journal of Machine Learning Research">JMLR</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Related articles</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Glossary_of_artificial_intelligence" title="Glossary of artificial intelligence">Glossary of artificial intelligence</a></li>
<li><a href="List_of_datasets_for_machine-learning_research" title="List of datasets for machine-learning research">List of datasets for machine-learning research</a>
<ul><li><a href="List_of_datasets_in_computer_vision_and_image_processing" title="List of datasets in computer vision and image processing">List of datasets in computer vision and image processing</a></li></ul></li>
<li><a href="Outline_of_machine_learning" title="Outline of machine learning">Outline of machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-navbar"><style data-mw-deduplicate="TemplateStyles:r1239400231">
/* start https://en.wikipedia.org/ */
.mw-parser-output .navbar{display:inline;font-size:88%;font-weight:normal}.mw-parser-output .navbar-collapse{float:left;text-align:left}.mw-parser-output .navbar-boxtext{word-spacing:0}.mw-parser-output .navbar ul{display:inline-block;white-space:nowrap;line-height:inherit}.mw-parser-output .navbar-brackets::before{margin-right:-0.125em;content:"[ "}.mw-parser-output .navbar-brackets::after{margin-left:-0.125em;content:" ]"}.mw-parser-output .navbar li{word-spacing:-0.125em}.mw-parser-output .navbar a>span,.mw-parser-output .navbar a>abbr{text-decoration:inherit}.mw-parser-output .navbar-mini abbr{font-variant:small-caps;border-bottom:none;text-decoration:none;cursor:inherit}.mw-parser-output .navbar-ct-full{font-size:114%;margin:0 7em}.mw-parser-output .navbar-ct-mini{font-size:114%;margin:0 4em}html.skin-theme-clientpref-night .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}@media(prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}}@media print{.mw-parser-output .navbar{display:none!important}}
/* end https://en.wikipedia.org/ */
</style></td></tr></tbody></table>
<p><b>Unsupervised learning</b> is a framework in <a href="Machine_learning" title="Machine learning">machine learning</a> where, in contrast to <a href="Supervised_learning" title="Supervised learning">supervised learning</a>, algorithms learn patterns exclusively from unlabeled data.<sup id="cite_ref-WeiWu_1-0" class="reference"><a href="#cite_note-WeiWu-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup> Other frameworks in the spectrum of supervisions include <a href="Weak_supervision" title="Weak supervision">weak- or semi-supervision</a>, where a small portion of the data is tagged, and <a href="Self-supervised_learning" title="Self-supervised learning">self-supervision</a>. Some researchers consider self-supervised learning a form of unsupervised learning.<sup id="cite_ref-2" class="reference"><a href="#cite_note-2"><span class="cite-bracket">[</span>2<span class="cite-bracket">]</span></a></sup>
</p><p>Conceptually, unsupervised learning divides into the aspects of data, training, algorithm, and downstream applications. Typically, the dataset is harvested cheaply "in the wild", such as massive <a href="Text_corpus" title="Text corpus">text corpus</a> obtained by <a href="Web_crawler" title="Web crawler">web crawling</a>, with only minor filtering (such as <a href="Common_Crawl" title="Common Crawl">Common Crawl</a>). This compares favorably to supervised learning, where the dataset (such as the <a href="ImageNet" title="ImageNet">ImageNet1000</a>) is typically constructed manually, which is much more expensive.
</p><p>There were algorithms designed specifically for unsupervised learning, such as <a href="Cluster_analysis" title="Cluster analysis">clustering algorithms</a> like <a href="K-means_clustering" title="K-means clustering">k-means</a>, <a href="Dimensionality_reduction" title="Dimensionality reduction">dimensionality reduction</a> techniques like <a href="Principal_component_analysis" title="Principal component analysis">principal component analysis (PCA)</a>, <a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann machine learning</a>, and <a href="Autoencoder" title="Autoencoder">autoencoders</a>. After the rise of deep learning, most large-scale unsupervised learning have been done by training general-purpose neural network architectures by <a href="Gradient_descent" title="Gradient descent">gradient descent</a>, adapted to performing unsupervised learning by designing an appropriate training procedure.
</p><p>Sometimes a trained model can be used as-is, but more often they are modified for downstream applications. For example, the generative pretraining method trains a model to generate a textual dataset, before finetuning it for other applications, such as text classification.<sup id="cite_ref-gpt1paper_3-0" class="reference"><a href="#cite_note-gpt1paper-3"><span class="cite-bracket">[</span>3<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-4" class="reference"><a href="#cite_note-4"><span class="cite-bracket">[</span>4<span class="cite-bracket">]</span></a></sup> As another example, autoencoders are trained to <a href="Feature_learning" title="Feature learning">good features</a>, which can then be used as a module for other models, such as in a <a href="Latent_diffusion_model" title="Latent diffusion model">latent diffusion model</a>.
</p>
<meta property="mw:PageProp/toc">
<div class="mw-heading mw-heading2"><h2 id="Tasks">Tasks</h2></div>
<p>Tasks are often categorized as <a href="Discriminative_model" title="Discriminative model">discriminative</a> (recognition) or <a href="Generative_model" title="Generative model">generative</a> (imagination). Often but not always, discriminative tasks use supervised methods and generative tasks use unsupervised (see <a href="Venn_diagram" title="Venn diagram">Venn diagram</a>); however, the separation is very hazy. For example, object recognition favors supervised learning but unsupervised learning can also cluster objects into groups. Furthermore, as progress marches onward, some tasks employ both methods, and some tasks swing from one to another. For example, image recognition started off as heavily supervised, but became hybrid by employing unsupervised pre-training, and then moved towards supervision again with the advent of <a href="Dilution_(neural_networks)" title="Dilution (neural networks)">dropout</a>, <a href="Rectifier_(neural_networks)" title="Rectifier (neural networks)">ReLU</a>, and <a href="Learning_rate" title="Learning rate">adaptive learning rates</a>.
</p><p>A typical generative task is as follows. At each step, a datapoint is sampled from the dataset, and part of the data is removed, and the model must infer the removed part. This is particularly clear for the <a href="Autoencoder" title="Autoencoder">denoising autoencoders</a> and <a href="BERT_(language_model)" title="BERT (language model)">BERT</a>.
</p>
<div class="mw-heading mw-heading2"><h2 id="Neural_network_architectures">Neural network architectures</h2></div>
<div class="mw-heading mw-heading3"><h3 id="Training">Training</h3></div>
<p>During the learning phase, an unsupervised network tries to mimic the data it's given and uses the error in its mimicked output to correct itself (i.e. correct its weights and biases). Sometimes the error is expressed as a low probability that the erroneous output occurs, or it might be expressed as an unstable high energy state in the network.
</p><p>In contrast to supervised methods' dominant use of <a href="Backpropagation" title="Backpropagation">backpropagation</a>, unsupervised learning also employs other methods including: Hopfield learning rule, Boltzmann learning rule, <a href="Contrastive_Divergence" class="mw-redirect" title="Contrastive Divergence">Contrastive Divergence</a>, <a href="Wake-sleep_algorithm" title="Wake-sleep algorithm">Wake Sleep</a>, <a href="Variational_Inference" class="mw-redirect" title="Variational Inference">Variational Inference</a>, <a href="Maximum_Likelihood" class="mw-redirect" title="Maximum Likelihood">Maximum Likelihood</a>, <a href="Maximum_A_Posteriori" class="mw-redirect" title="Maximum A Posteriori">Maximum A Posteriori</a>, <a href="Gibbs_Sampling" class="mw-redirect" title="Gibbs Sampling">Gibbs Sampling</a>, and backpropagating reconstruction errors or hidden state reparameterizations. See the table below for more details.
</p>
<div class="mw-heading mw-heading3"><h3 id="Energy">Energy</h3></div>
<p>An energy function is a macroscopic measure of a network's activation state. In Boltzmann machines, it plays the role of the Cost function. This analogy with physics is inspired by Ludwig Boltzmann's analysis of a gas' macroscopic energy from the microscopic probabilities of particle motion <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle p\propto e^{-E/kT}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>p</mi>
<mo>∝<!-- ∝ --></mo>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo>−<!-- − --></mo>
<mi>E</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo>/</mo>
</mrow>
<mi>k</mi>
<mi>T</mi>
</mrow>
</msup>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle p\propto e^{-E/kT}}</annotation>
</semantics>
</math></span><img src="./b9696fac92eaa5fede297c17fa8bb086945efd8e.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; margin-left: -0.089ex; width:11.043ex; height:3.176ex;" alt="{\displaystyle p\propto e^{-E/kT}}" loading="lazy"></span>, where k is the Boltzmann constant and T is temperature. In the <a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">RBM</a> network the relation is <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle p=e^{-E}/Z}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>p</mi>
<mo>=</mo>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo>−<!-- − --></mo>
<mi>E</mi>
</mrow>
</msup>
<mrow class="MJX-TeXAtom-ORD">
<mo>/</mo>
</mrow>
<mi>Z</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle p=e^{-E}/Z}</annotation>
</semantics>
</math></span><img src="./7c5acd1a54ea1654f16ad3b413693ab78c72319a.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; margin-left: -0.089ex; width:11.05ex; height:3.176ex;" alt="{\displaystyle p=e^{-E}/Z}" loading="lazy"></span>,<sup id="cite_ref-Hinton2010_5-0" class="reference"><a href="#cite_note-Hinton2010-5"><span class="cite-bracket">[</span>5<span class="cite-bracket">]</span></a></sup> where <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle p}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>p</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle p}</annotation>
</semantics>
</math></span><img src="./81eac1e205430d1f40810df36a0edffdc367af36.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; margin-left: -0.089ex; width:1.259ex; height:2.009ex;" alt="{\displaystyle p}" loading="lazy"></span> and <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle E}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>E</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle E}</annotation>
</semantics>
</math></span><img src="./4232c9de2ee3eec0a9c0a19b15ab92daa6223f9b.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.776ex; height:2.176ex;" alt="{\displaystyle E}" loading="lazy"></span> vary over every possible activation pattern and <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \textstyle {Z=\sum _{\scriptscriptstyle {\text{All Patterns}}}e^{-E({\text{pattern}})}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mstyle displaystyle="false" scriptlevel="0">
<mrow class="MJX-TeXAtom-ORD">
<mi>Z</mi>
<mo>=</mo>
<munder>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="false" scriptlevel="2">
<mrow class="MJX-TeXAtom-ORD">
<mtext>All Patterns</mtext>
</mrow>
</mstyle>
</mrow>
</munder>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo>−<!-- − --></mo>
<mi>E</mi>
<mo stretchy="false">(</mo>
<mrow class="MJX-TeXAtom-ORD">
<mtext>pattern</mtext>
</mrow>
<mo stretchy="false">)</mo>
</mrow>
</msup>
</mrow>
</mstyle>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \textstyle {Z=\sum _{\scriptscriptstyle {\text{All Patterns}}}e^{-E({\text{pattern}})}}}</annotation>
</semantics>
</math></span><img src="./b2d124436452bf4f7c46be05f2a4ec154e266c2d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -1.005ex; width:25.608ex; height:3.343ex;" alt="{\displaystyle \textstyle {Z=\sum _{\scriptscriptstyle {\text{All Patterns}}}e^{-E({\text{pattern}})}}}" loading="lazy"></span>. To be more precise, <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle p(a)=e^{-E(a)}/Z}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>p</mi>
<mo stretchy="false">(</mo>
<mi>a</mi>
<mo stretchy="false">)</mo>
<mo>=</mo>
<msup>
<mi>e</mi>
<mrow class="MJX-TeXAtom-ORD">
<mo>−<!-- − --></mo>
<mi>E</mi>
<mo stretchy="false">(</mo>
<mi>a</mi>
<mo stretchy="false">)</mo>
</mrow>
</msup>
<mrow class="MJX-TeXAtom-ORD">
<mo>/</mo>
</mrow>
<mi>Z</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle p(a)=e^{-E(a)}/Z}</annotation>
</semantics>
</math></span><img src="./749ebbae5c41727f56fbb5fbee29d6f792ccfa4f.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; margin-left: -0.089ex; width:16.238ex; height:3.343ex;" alt="{\displaystyle p(a)=e^{-E(a)}/Z}" loading="lazy"></span>, where <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle a}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>a</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle a}</annotation>
</semantics>
</math></span><img src="./ffd2487510aa438433a2579450ab2b3d557e5edc.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.23ex; height:1.676ex;" alt="{\displaystyle a}" loading="lazy"></span> is an activation pattern of all neurons (visible and hidden). Hence, some early neural networks bear the name Boltzmann Machine. Paul Smolensky calls <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle -E\,}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo>−<!-- − --></mo>
<mi>E</mi>
<mspace width="thinmathspace"></mspace>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle -E\,}</annotation>
</semantics>
</math></span><img src="./27723660512942ff84ce2d85df2509299c698147.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.505ex; width:3.971ex; height:2.343ex;" alt="{\displaystyle -E\,}" loading="lazy"></span> the <i>Harmony</i>. A network seeks low energy which is high Harmony.
</p>
<div class="mw-heading mw-heading3"><h3 id="Networks">Networks</h3></div>
<p>This table shows connection diagrams of various unsupervised networks, the details of which will be given in the section Comparison of Networks. Circles are neurons and edges between them are connection weights. As network design changes, features are added on to enable new capabilities or removed to make learning faster. For instance, neurons change between deterministic (Hopfield) and stochastic (Boltzmann) to allow robust output, weights are removed within a layer (RBM) to hasten learning, or connections are allowed to become asymmetric (Helmholtz).
</p>
<table class="wikitable">
<tbody><tr>
<th><a href="Hopfield_network" title="Hopfield network">Hopfield</a></th>
<th><a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann</a></th>
<th><a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">RBM</a></th>
<th><a href="Stacked_Restricted_Boltzmann_Machine" class="mw-redirect" title="Stacked Restricted Boltzmann Machine">Stacked Boltzmann</a>
</th></tr>
<tr>
<td>
</td>
<td>
</td>
<td>
</td>
<td>
</td></tr></tbody></table>
<table class="wikitable">
<tbody><tr>
<th><a href="Helmholtz_machine" title="Helmholtz machine">Helmholtz</a></th>
<th><a href="Autoencoder" title="Autoencoder">Autoencoder</a></th>
<th><a href="Variational_autoencoder" title="Variational autoencoder">VAE</a>
</th></tr>
<tr>
<td>
</td>
<td>
</td>
<td>
</td></tr></tbody></table>
<p>Of the networks bearing people's names, only Hopfield worked directly with neural networks. Boltzmann and Helmholtz came before artificial neural networks, but their work in physics and physiology inspired the analytical methods that were used.
</p>
<div class="mw-heading mw-heading3"><h3 id="History">History</h3></div>
<table class="wikitable">
<tbody><tr>
<td>1974</td>
<td>Ising magnetic model proposed by WA Little for cognition
</td></tr>
<tr>
<td>1980</td>
<td><a href="Kunihiko_Fukushima" title="Kunihiko Fukushima">Kunihiko Fukushima</a> introduces the <a href="Neocognitron" title="Neocognitron">neocognitron</a>, which is later called a <a href="Convolutional_neural_network" title="Convolutional neural network">convolutional neural network</a>. It is mostly used in SL, but deserves a mention here.
</td></tr>
<tr>
<td>1982</td>
<td>Ising variant Hopfield net described as <a href="Content-addressable_memory" title="Content-addressable memory">CAMs</a> and classifiers by John Hopfield.
</td></tr>
<tr>
<td>1983</td>
<td>Ising variant Boltzmann machine with probabilistic neurons described by <a href="Geoffrey_Hinton" title="Geoffrey Hinton">Hinton</a> & <a href="Terry_Sejnowski" title="Terry Sejnowski">Sejnowski</a> following Sherington & Kirkpatrick's 1975 work.
</td></tr>
<tr>
<td>1986</td>
<td><a href="Paul_Smolensky" title="Paul Smolensky">Paul Smolensky</a> publishes Harmony Theory, which is an RBM with practically the same Boltzmann energy function. Smolensky did not give a practical training scheme. Hinton did in mid-2000s.
</td></tr>
<tr>
<td>1995</td>
<td>Schmidthuber introduces the <a href="Long_short-term_memory" title="Long short-term memory">LSTM</a> neuron for languages.
</td></tr>
<tr>
<td>1995</td>
<td>Dayan & Hinton introduces Helmholtz machine
</td></tr>
<tr>
<td>2013</td>
<td>Kingma, Rezende, & co. introduced Variational Autoencoders as Bayesian graphical probability network, with neural nets as components.
</td></tr></tbody></table>
<div class="mw-heading mw-heading3"><h3 id="Specific_Networks">Specific Networks</h3></div>
<p>Here, we highlight some characteristics of select networks. The details of each are given in the comparison table below.
</p>
<style data-mw-deduplicate="TemplateStyles:r1228772891">
/* start https://en.wikipedia.org/ */
.mw-parser-output .glossary dt{margin-top:0.4em}.mw-parser-output .glossary dt+dt{margin-top:-0.2em}.mw-parser-output .glossary .templatequote{margin-top:0;margin-bottom:-0.5em}
/* end https://en.wikipedia.org/ */
</style>
<dl class="glossary">
<dt id="hopfield_network"><dfn><a href="Hopfield_Network" class="mw-redirect" title="Hopfield Network">Hopfield Network</a></dfn></dt>
<dd>Ferromagnetism inspired Hopfield networks. A neuron correspond to an iron domain with binary magnetic moments Up and Down, and neural connections correspond to the domain's influence on each other. Symmetric connections enable a global energy formulation. During inference the network updates each state using the standard activation step function. Symmetric weights and the right energy functions guarantees convergence to a stable activation pattern. Asymmetric weights are difficult to analyze. Hopfield nets are used as Content Addressable Memories (CAM).</dd>
<dt id="boltzmann_machine"><dfn><a href="Boltzmann_Machine" class="mw-redirect" title="Boltzmann Machine">Boltzmann Machine</a></dfn></dt>
<dd>These are stochastic Hopfield nets. Their state value is sampled from this <a href="Probability_density_function" title="Probability density function">pdf</a> as follows: suppose a binary neuron fires with the Bernoulli probability p(1) = 1/3 and rests with p(0) = 2/3. One samples from it by taking a <i>uniformly</i> distributed random number y, and plugging it into the inverted <a href="Cumulative_distribution_function" title="Cumulative distribution function">cumulative distribution function</a>, which in this case is the step function thresholded at 2/3. The inverse function = { 0 if x <= 2/3, 1 if x > 2/3 }.</dd>
<dt id="sigmoid_belief_net"><dfn>Sigmoid Belief Net</dfn></dt>
<dd>Introduced by Radford Neal in 1992, this network applies ideas from probabilistic graphical models to neural networks. A key difference is that nodes in graphical models have pre-assigned meanings, whereas Belief Net neurons' features are determined after training. The network is a sparsely connected directed acyclic graph composed of binary stochastic neurons. The learning rule comes from Maximum Likelihood on p(X): Δw<sub>ij</sub> <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle \propto }">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mo>∝<!-- ∝ --></mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle \propto }</annotation>
</semantics>
</math></span><img src="./0e3a55007ba2f092d6cafe6d33598e0608b81150.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.808ex; height:1.676ex;" alt="{\displaystyle \propto }" loading="lazy"></span> s<sub>j</sub> * (s<sub>i</sub> - p<sub>i</sub>), where p<sub>i</sub> = 1 / ( 1 + e<sup>weighted inputs into neuron i</sup> ). s<sub>j</sub>'s are activations from an unbiased sample of the posterior distribution and this is problematic due to the Explaining Away problem raised by Judea Perl. <a href="Variational_Bayesian_methods" title="Variational Bayesian methods">Variational Bayesian methods</a> uses a surrogate posterior and blatantly disregard this complexity.</dd>
<dt id="deep_belief_network"><dfn><a href="Deep_Belief_Network" class="mw-redirect" title="Deep Belief Network">Deep Belief Network</a></dfn></dt>
<dd>Introduced by Hinton, this network is a hybrid of RBM and Sigmoid Belief Network. The top 2 layers is an RBM and the second layer downwards form a sigmoid belief network. One trains it by the <a href="Stacked_Restricted_Boltzmann_Machine" class="mw-redirect" title="Stacked Restricted Boltzmann Machine">stacked RBM</a> method and then throw away the recognition weights below the top RBM. As of 2009, 3-4 layers seems to be the optimal depth.<sup id="cite_ref-HintonMlss2009_6-0" class="reference"><a href="#cite_note-HintonMlss2009-6"><span class="cite-bracket">[</span>6<span class="cite-bracket">]</span></a></sup></dd>
<dt id="helmholtz_machine"><dfn><a href="Helmholtz_machine" title="Helmholtz machine">Helmholtz machine</a></dfn></dt>
<dd>These are early inspirations for the Variational Auto Encoders. Its 2 networks combined into one—forward weights operates recognition and backward weights implements imagination. It is perhaps the first network to do both. Helmholtz did not work in machine learning but he inspired the view of "statistical inference engine whose function is to infer probable causes of sensory input".<sup id="cite_ref-nc95_7-0" class="reference"><a href="#cite_note-nc95-7"><span class="cite-bracket">[</span>7<span class="cite-bracket">]</span></a></sup> the stochastic binary neuron outputs a probability that its state is 0 or 1. The data input is normally not considered a layer, but in the Helmholtz machine generation mode, the data layer receives input from the middle layer and has separate weights for this purpose, so it is considered a layer. Hence this network has 3 layers.</dd>
<dt id="variational_autoencoder"><dfn><a href="Variational_autoencoder" title="Variational autoencoder">Variational autoencoder</a></dfn></dt>
<dd>These are inspired by Helmholtz machines and combines probability network with neural networks. An Autoencoder is a 3-layer CAM network, where the middle layer is supposed to be some internal representation of input patterns. The encoder neural network is a probability distribution q<sub>φ</sub>(z given x) and the decoder network is p<sub>θ</sub>(x given z). The weights are named phi & theta rather than W and V as in Helmholtz—a cosmetic difference. These 2 networks here can be fully connected, or use another NN scheme.</dd>
</dl>
<div class="mw-heading mw-heading3"><h3 id="Comparison_of_networks">Comparison of networks</h3></div>
<table class="wikitable">
<tbody><tr>
<th></th>
<th>Hopfield</th>
<th>Boltzmann</th>
<th>RBM</th>
<th>Stacked RBM</th>
<th>Helmholtz</th>
<th>Autoencoder</th>
<th>VAE
</th></tr>
<tr>
<td><b>Usage & notables</b></td>
<td>CAM, traveling salesman problem</td>
<td>CAM. The freedom of connections makes this network difficult to analyze.</td>
<td>pattern recognition. used in MNIST digits and speech.</td>
<td>recognition & imagination. trained with unsupervised pre-training and/or supervised fine tuning.</td>
<td>imagination, mimicry</td>
<td>language: creative writing, translation. vision: enhancing blurry images</td>
<td>generate realistic data
</td></tr>
<tr>
<td><b>Neuron</b></td>
<td>deterministic binary state. Activation = { 0 (or -1) if x is negative, 1 otherwise }</td>
<td>stochastic binary Hopfield neuron</td>
<td>← same. (extended to real-valued in mid 2000s)</td>
<td>← same</td>
<td>← same</td>
<td>language: LSTM. vision: local receptive fields. usually real valued relu activation.</td>
<td>middle layer neurons encode means & variances for Gaussians. In run mode (inference), the output of the middle layer are sampled values from the Gaussians.
</td></tr>
<tr>
<td><b>Connections</b></td>
<td>1-layer with symmetric weights. No self-connections.</td>
<td>2-layers. 1-hidden & 1-visible. symmetric weights.</td>
<td>← same. <br>no lateral connections within a layer.</td>
<td>top layer is undirected, symmetric. other layers are 2-way, asymmetric.</td>
<td>3-layers: asymmetric weights. 2 networks combined into 1.</td>
<td>3-layers. The input is considered a layer even though it has no inbound weights. recurrent layers for NLP. feedforward convolutions for vision. input & output have the same neuron counts.</td>
<td>3-layers: input, encoder, distribution sampler decoder. the sampler is not considered a layer
</td></tr>
<tr>
<td><b>Inference & energy</b></td>
<td>Energy is given by Gibbs probability measure :<span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle E=-{\frac {1}{2}}\sum _{i,j}{w_{ij}{s_{i}}{s_{j}}}+\sum _{i}{\theta _{i}}{s_{i}}}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>E</mi>
<mo>=</mo>
<mo>−<!-- − --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mfrac>
<mn>1</mn>
<mn>2</mn>
</mfrac>
</mrow>
<munder>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mo>,</mo>
<mi>j</mi>
</mrow>
</munder>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>w</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
<mi>j</mi>
</mrow>
</msub>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>j</mi>
</mrow>
</msub>
</mrow>
</mrow>
<mo>+</mo>
<munder>
<mo>∑<!-- ∑ --></mo>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</munder>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>θ<!-- θ --></mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
<mrow class="MJX-TeXAtom-ORD">
<msub>
<mi>s</mi>
<mrow class="MJX-TeXAtom-ORD">
<mi>i</mi>
</mrow>
</msub>
</mrow>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle E=-{\frac {1}{2}}\sum _{i,j}{w_{ij}{s_{i}}{s_{j}}}+\sum _{i}{\theta _{i}}{s_{i}}}</annotation>
</semantics>
</math></span><img src="./76ad57f7823ca1b74ab67b5f3dadbc83d24da7c2.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -3.338ex; width:30.204ex; height:6.676ex;" alt="{\displaystyle E=-{\frac {1}{2}}\sum _{i,j}{w_{ij}{s_{i}}{s_{j}}}+\sum _{i}{\theta _{i}}{s_{i}}}" loading="lazy"></span></td>
<td>← same</td>
<td>← same</td>
<td></td>
<td>minimize KL divergence</td>
<td>inference is only feed-forward. previous UL networks ran forwards AND backwards</td>
<td>minimize error = reconstruction error - KLD
</td></tr>
<tr>
<td><b>Training</b></td>
<td>Δw<sub>ij</sub> = s<sub>i</sub>*s<sub>j</sub>, for +1/-1 neuron</td>
<td>Δw<sub>ij</sub> = e*(p<sub>ij</sub> - p'<sub>ij</sub>). This is derived from minimizing KLD. e = learning rate, p' = predicted and p = actual distribution.
</td>
<td>Δw<sub>ij</sub> = e*( < v<sub>i</sub> h<sub>j</sub> ><sub>data</sub> - < v<sub>i</sub> h<sub>j</sub> ><sub>equilibrium</sub> ). This is a form of contrastive divergence w/ Gibbs Sampling. "<>" are expectations.</td>
<td>← similar. train 1-layer at a time. approximate equilibrium state with a 3-segment pass. no back propagation.</td>
<td>wake-sleep 2 phase training</td>
<td>back propagate the reconstruction error</td>
<td>reparameterize hidden state for backprop
</td></tr>
<tr>
<td><b>Strength</b></td>
<td>resembles physical systems so it inherits their equations</td>
<td>← same. hidden neurons act as internal representatation of the external world</td>
<td>faster more practical training scheme than Boltzmann machines</td>
<td>trains quickly. gives hierarchical layer of features</td>
<td>mildly anatomical. analyzable w/ information theory & statistical mechanics</td>
<td></td>
<td>
</td></tr>
<tr>
<td><b>Weakness</b></td>
<td></td>
<td>hard to train due to lateral connections</td>
<td>equilibrium requires too many iterations</td>
<td>integer & real-valued neurons are more complicated.</td>
<td></td>
<td></td>
<td>
</td></tr></tbody></table>
<div class="mw-heading mw-heading3"><h3 id="Hebbian_Learning,_ART,_SOM">Hebbian Learning, ART, SOM</h3></div>
<p>The classical example of unsupervised learning in the study of neural networks is <a href="Donald_Hebb" class="mw-redirect" title="Donald Hebb">Donald Hebb</a>'s principle, that is, neurons that fire together wire together.<sup id="cite_ref-Buhmann_8-0" class="reference"><a href="#cite_note-Buhmann-8"><span class="cite-bracket">[</span>8<span class="cite-bracket">]</span></a></sup> In <a href="Hebbian_learning" class="mw-redirect" title="Hebbian learning">Hebbian learning</a>, the connection is reinforced irrespective of an error, but is exclusively a function of the coincidence between action potentials between the two neurons.<sup id="cite_ref-Comesana_9-0" class="reference"><a href="#cite_note-Comesana-9"><span class="cite-bracket">[</span>9<span class="cite-bracket">]</span></a></sup> A similar version that modifies synaptic weights takes into account the time between the action potentials (<a href="Spike-timing-dependent_plasticity" title="Spike-timing-dependent plasticity">spike-timing-dependent plasticity</a> or STDP). Hebbian Learning has been hypothesized to underlie a range of cognitive functions, such as <a href="Pattern_recognition" title="Pattern recognition">pattern recognition</a> and experiential learning.
</p><p>Among <a href="Artificial_neural_network" class="mw-redirect" title="Artificial neural network">neural network</a> models, the <a href="Self-organizing_map" title="Self-organizing map">self-organizing map</a> (SOM) and <a href="Adaptive_resonance_theory" title="Adaptive resonance theory">adaptive resonance theory</a> (ART) are commonly used in unsupervised learning algorithms. The SOM is a topographic organization in which nearby locations in the map represent inputs with similar properties. The ART model allows the number of clusters to vary with problem size and lets the user control the degree of similarity between members of the same clusters by means of a user-defined constant called the vigilance parameter. ART networks are used for many pattern recognition tasks, such as <a href="Automatic_target_recognition" title="Automatic target recognition">automatic target recognition</a> and seismic signal processing.<sup id="cite_ref-Carpenter_10-0" class="reference"><a href="#cite_note-Carpenter-10"><span class="cite-bracket">[</span>10<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Probabilistic_methods">Probabilistic methods</h2></div>
<p>Two of the main methods used in unsupervised learning are <a href="Principal_component_analysis" title="Principal component analysis">principal component</a> and <a href="Cluster_analysis" title="Cluster analysis">cluster analysis</a>. Cluster analysis is used in unsupervised learning to group, or segment, datasets with shared attributes in order to extrapolate algorithmic relationships.<sup id="cite_ref-tds-ul_11-0" class="reference"><a href="#cite_note-tds-ul-11"><span class="cite-bracket">[</span>11<span class="cite-bracket">]</span></a></sup> Cluster analysis is a branch of <a href="Machine_learning" title="Machine learning">machine learning</a> that groups the data that has not been <a href="Labeled_data" title="Labeled data">labelled</a>, classified or categorized. Instead of responding to feedback, cluster analysis identifies commonalities in the data and reacts based on the presence or absence of such commonalities in each new piece of data. This approach helps detect anomalous data points that do not fit into either group.
</p><p>A central application of unsupervised learning is in the field of <a href="Density_estimation" title="Density estimation">density estimation</a> in <a href="Statistics" title="Statistics">statistics</a>,<sup id="cite_ref-JordanBishop2004_12-0" class="reference"><a href="#cite_note-JordanBishop2004-12"><span class="cite-bracket">[</span>12<span class="cite-bracket">]</span></a></sup> though unsupervised learning encompasses many other domains involving summarizing and explaining data features. It can be contrasted with supervised learning by saying that whereas supervised learning intends to infer a <a href="Conditional_probability_distribution" title="Conditional probability distribution">conditional probability distribution</a> conditioned on the label of input data; unsupervised learning intends to infer an <a href="A_priori_probability" class="mw-redirect" title="A priori probability">a priori probability</a> distribution .
</p>
<div class="mw-heading mw-heading3"><h3 id="Approaches">Approaches</h3></div>
<p>Some of the most common algorithms used in unsupervised learning include: (1) Clustering, (2) Anomaly detection, (3) Approaches for learning latent variable models. Each approach uses several methods as follows:
</p>
<ul><li><a href="Data_clustering" class="mw-redirect" title="Data clustering">Clustering</a> methods include: <a href="Hierarchical_clustering" title="Hierarchical clustering">hierarchical clustering</a>,<sup id="cite_ref-Hastie_13-0" class="reference"><a href="#cite_note-Hastie-13"><span class="cite-bracket">[</span>13<span class="cite-bracket">]</span></a></sup> <a href="K-means" class="mw-redirect" title="K-means">k-means</a>,<sup id="cite_ref-tds-kmeans_14-0" class="reference"><a href="#cite_note-tds-kmeans-14"><span class="cite-bracket">[</span>14<span class="cite-bracket">]</span></a></sup> <a href="Mixture_models" class="mw-redirect" title="Mixture models">mixture models</a>, <a href="Model-based_clustering" title="Model-based clustering">model-based clustering</a>, <a href="DBSCAN" title="DBSCAN">DBSCAN</a>, and <a href="OPTICS_algorithm" title="OPTICS algorithm">OPTICS algorithm</a></li>
<li><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a> methods include: <a href="Local_Outlier_Factor" class="mw-redirect" title="Local Outlier Factor">Local Outlier Factor</a>, and <a href="Isolation_Forest" class="mw-redirect" title="Isolation Forest">Isolation Forest</a></li>
<li>Approaches for learning <a href="Latent_variable_model" title="Latent variable model">latent variable models</a> such as <a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization algorithm</a> (EM), <a href="Method_of_moments_(statistics)" title="Method of moments (statistics)">Method of moments</a>, and <a href="Blind_signal_separation" class="mw-redirect" title="Blind signal separation">Blind signal separation</a> techniques (Principal component analysis, <a href="Independent_component_analysis" title="Independent component analysis">Independent component analysis</a>, <a href="Non-negative_matrix_factorization" title="Non-negative matrix factorization">Non-negative matrix factorization</a>, <a href="Singular_value_decomposition" title="Singular value decomposition">Singular value decomposition</a>)</li></ul>
<div class="mw-heading mw-heading3"><h3 id="Method_of_moments">Method of moments</h3></div>
<p>One of the statistical approaches for unsupervised learning is the <a href="Method_of_moments_(statistics)" title="Method of moments (statistics)">method of moments</a>. In the method of moments, the unknown parameters (of interest) in the model are related to the moments of one or more random variables, and thus, these unknown parameters can be estimated given the moments. The moments are usually estimated from samples empirically. The basic moments are first and second order moments. For a random vector, the first order moment is the <a href="Mean" title="Mean">mean</a> vector, and the second order moment is the <a href="Covariance_matrix" title="Covariance matrix">covariance matrix</a> (when the mean is zero). Higher order moments are usually represented using <a href="Tensors" class="mw-redirect" title="Tensors">tensors</a> which are the generalization of matrices to higher orders as multi-dimensional arrays.
</p><p>In particular, the method of moments is shown to be effective in learning the parameters of <a href="Latent_variable_model" title="Latent variable model">latent variable models</a>. Latent variable models are statistical models where in addition to the observed variables, a set of latent variables also exists which is not observed. A highly practical example of latent variable models in machine learning is the <a href="Topic_modeling" class="mw-redirect" title="Topic modeling">topic modeling</a> which is a statistical model for generating the words (observed variables) in the document based on the topic (latent variable) of the document. In the topic modeling, the words in the document are generated according to different statistical parameters when the topic of the document is changed. It is shown that method of moments (tensor decomposition techniques) consistently recover the parameters of a large class of latent variable models under some assumptions.<sup id="cite_ref-TensorLVMs_15-0" class="reference"><a href="#cite_note-TensorLVMs-15"><span class="cite-bracket">[</span>15<span class="cite-bracket">]</span></a></sup>
</p><p>The <a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization algorithm</a> (EM) is also one of the most practical methods for learning latent variable models. However, it can get stuck in local optima, and it is not guaranteed that the algorithm will converge to the true unknown parameters of the model. In contrast, for the method of moments, the global convergence is guaranteed under some conditions.
</p>
<div class="mw-heading mw-heading2"><h2 id="See_also">See also</h2></div>
<ul><li><a href="Automated_machine_learning" title="Automated machine learning">Automated machine learning</a></li>
<li><a href="Cluster_analysis" title="Cluster analysis">Cluster analysis</a></li>
<li><a href="Model-based_clustering" title="Model-based clustering">Model-based clustering</a></li>
<li><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></li>
<li><a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization algorithm</a></li>
<li><a href="Generative_topographic_map" title="Generative topographic map">Generative topographic map</a></li>
<li><a href="Meta-learning_(computer_science)" title="Meta-learning (computer science)">Meta-learning (computer science)</a></li>
<li><a href="Multivariate_analysis" class="mw-redirect" title="Multivariate analysis">Multivariate analysis</a></li>
<li><a href="Radial_basis_function_network" title="Radial basis function network">Radial basis function network</a></li>
<li><a href="Weak_supervision" title="Weak supervision">Weak supervision</a></li></ul>
<div class="mw-heading mw-heading2"><h2 id="References">References</h2></div>
<style data-mw-deduplicate="TemplateStyles:r1239543626">
/* start https://en.wikipedia.org/ */
.mw-parser-output .reflist{margin-bottom:0.5em;list-style-type:decimal}@media screen{.mw-parser-output .reflist{font-size:90%}}.mw-parser-output .reflist .references{font-size:100%;margin-bottom:0;list-style-type:inherit}.mw-parser-output .reflist-columns-2{column-width:30em}.mw-parser-output .reflist-columns-3{column-width:25em}.mw-parser-output .reflist-columns{margin-top:0.3em}.mw-parser-output .reflist-columns ol{margin-top:0}.mw-parser-output .reflist-columns li{page-break-inside:avoid;break-inside:avoid-column}.mw-parser-output .reflist-upper-alpha{list-style-type:upper-alpha}.mw-parser-output .reflist-upper-roman{list-style-type:upper-roman}.mw-parser-output .reflist-lower-alpha{list-style-type:lower-alpha}.mw-parser-output .reflist-lower-greek{list-style-type:lower-greek}.mw-parser-output .reflist-lower-roman{list-style-type:lower-roman}
/* end https://en.wikipedia.org/ */
</style><div class="reflist">
<div class="mw-references-wrap mw-references-columns"><ol class="references">
<li id="cite_note-WeiWu-1"><span class="mw-cite-backlink"><b><a href="#cite_ref-WeiWu_1-0">^</a></b></span> <span class="reference-text"><style data-mw-deduplicate="TemplateStyles:r1238218222">
/* start https://en.wikipedia.org/ */
.mw-parser-output cite.citation{font-style:inherit;word-wrap:break-word}.mw-parser-output .citation q{quotes:"\"""\"""'""'"}.mw-parser-output .citation:target{background-color:rgba(0,127,255,0.133)}.mw-parser-output .id-lock-free.id-lock-free a{background:url("./mw/Lock-green.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-limited.id-lock-limited a,.mw-parser-output .id-lock-registration.id-lock-registration a{background:url("./mw/Lock-gray-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-subscription.id-lock-subscription a{background:url("./mw/Lock-red-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .cs1-ws-icon a{background:url("./mw/Wikisource-logo.svg")right 0.1em center/12px no-repeat}body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-free a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-limited a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-registration a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-subscription a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .cs1-ws-icon a{background-size:contain;padding:0 1em 0 0}.mw-parser-output .cs1-code{color:inherit;background:inherit;border:none;padding:inherit}.mw-parser-output .cs1-hidden-error{display:none;color:var(--color-error,#d33)}.mw-parser-output .cs1-visible-error{color:var(--color-error,#d33)}.mw-parser-output .cs1-maint{display:none;color:#085;margin-left:0.3em}.mw-parser-output .cs1-kern-left{padding-left:0.2em}.mw-parser-output .cs1-kern-right{padding-right:0.2em}.mw-parser-output .citation .mw-selflink{font-weight:inherit}@media screen{.mw-parser-output .cs1-format{font-size:95%}html.skin-theme-clientpref-night .mw-parser-output .cs1-maint{color:#18911f}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .cs1-maint{color:#18911f}}
/* end https://en.wikipedia.org/ */
</style><cite id="CITEREFWu" class="citation web cs1">Wu, Wei. <a rel="nofollow" class="external text" href="https://na.uni-tuebingen.de/ex/ml_seminar_ss2022/Unsupervised_Learning%20Final.pdf">"Unsupervised Learning"</a> <span class="cs1-format">(PDF)</span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20240414213810/https://na.uni-tuebingen.de/ex/ml_seminar_ss2022/Unsupervised_Learning%20Final.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 14 April 2024<span class="reference-accessdate">. Retrieved <span class="nowrap">26 April</span> 2024</span>.</cite></span>
</li>
<li id="cite_note-2"><span class="mw-cite-backlink"><b><a href="#cite_ref-2">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiuZhangHouMian2021" class="citation journal cs1">Liu, Xiao; Zhang, Fanjin; Hou, Zhenyu; Mian, Li; Wang, Zhaoyu; Zhang, Jing; Tang, Jie (2021). <a rel="nofollow" class="external text" href="https://ieeexplore.ieee.org/document/9462394">"Self-supervised Learning: Generative or Contrastive"</a>. <i>IEEE Transactions on Knowledge and Data Engineering</i>: 1. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2006.08218">2006.08218</a></span>. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1109%2FTKDE.2021.3090866">10.1109/TKDE.2021.3090866</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a> <a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/1041-4347">1041-4347</a>.</cite></span>
</li>
<li id="cite_note-gpt1paper-3"><span class="mw-cite-backlink"><b><a href="#cite_ref-gpt1paper_3-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFRadfordNarasimhanSalimansSutskever2018" class="citation web cs1">Radford, Alec; Narasimhan, Karthik; Salimans, Tim; Sutskever, Ilya (11 June 2018). <a rel="nofollow" class="external text" href="https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf">"Improving Language Understanding by Generative Pre-Training"</a> <span class="cs1-format">(PDF)</span>. <a href="OpenAI" title="OpenAI">OpenAI</a>. p. 12. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20210126024542/https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 26 January 2021<span class="reference-accessdate">. Retrieved <span class="nowrap">23 January</span> 2021</span>.</cite></span>
</li>
<li id="cite_note-4"><span class="mw-cite-backlink"><b><a href="#cite_ref-4">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiWallaceShenLin2020" class="citation journal cs1">Li, Zhuohan; Wallace, Eric; Shen, Sheng; Lin, Kevin; Keutzer, Kurt; Klein, Dan; Gonzalez, Joey (2020-11-21). <a rel="nofollow" class="external text" href="https://proceedings.mlr.press/v119/li20m.html">"Train Big, Then Compress: Rethinking Model Size for Efficient Training and Inference of Transformers"</a>. <i>Proceedings of the 37th International Conference on Machine Learning</i>. PMLR: <span class="nowrap">5958–</span>5968.</cite></span>
</li>
<li id="cite_note-Hinton2010-5"><span class="mw-cite-backlink"><b><a href="#cite_ref-Hinton2010_5-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFHinton2012" class="citation book cs1">Hinton, G. (2012). <a rel="nofollow" class="external text" href="http://www.cs.utoronto.ca/~hinton/absps/guideTR.pdf">"A Practical Guide to Training Restricted Boltzmann Machines"</a> <span class="cs1-format">(PDF)</span>. <i>Neural Networks: Tricks of the Trade</i>. Lecture Notes in Computer Science. Vol. 7700. Springer. pp. <span class="nowrap">599–</span>619. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1007%2F978-3-642-35289-8_32">10.1007/978-3-642-35289-8_32</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>978-3-642-35289-8</bdi>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20220903215809/http://www.cs.utoronto.ca/~hinton/absps/guideTR.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2022-09-03<span class="reference-accessdate">. Retrieved <span class="nowrap">2022-11-03</span></span>.</cite></span>
</li>
<li id="cite_note-HintonMlss2009-6"><span class="mw-cite-backlink"><b><a href="#cite_ref-HintonMlss2009_6-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://videolectures.net/mlss09uk_hinton_dbn">"Deep Belief Nets"</a> (video). September 2009. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20220308022539/http://videolectures.net/mlss09uk_hinton_dbn/">Archived</a> from the original on 2022-03-08<span class="reference-accessdate">. Retrieved <span class="nowrap">2022-03-27</span></span>.</cite></span>
</li>
<li id="cite_note-nc95-7"><span class="mw-cite-backlink"><b><a href="#cite_ref-nc95_7-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFPeterHintonNealZemel1995" class="citation journal cs1"><a href="Peter_Dayan" title="Peter Dayan">Peter, Dayan</a>; <a href="Geoffrey_Hinton" title="Geoffrey Hinton">Hinton, Geoffrey E.</a>; <a href="Radford_M._Neal" title="Radford M. Neal">Neal, Radford M.</a>; <a href="Richard_Zemel" title="Richard Zemel">Zemel, Richard S.</a> (1995). "The Helmholtz machine". <i>Neural Computation</i>. <b>7</b> (5): <span class="nowrap">889–</span>904. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1162%2Fneco.1995.7.5.889">10.1162/neco.1995.7.5.889</a>. <a href="Hdl_(identifier)" class="mw-redirect" title="Hdl (identifier)">hdl</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://hdl.handle.net/21.11116%2F0000-0002-D6D3-E">21.11116/0000-0002-D6D3-E</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a> <a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/7584891">7584891</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a> <a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:1890561">1890561</a>.</cite> <span style="position:relative; top: -2px;"><span typeof="mw:File"><a href="Paywall" title="closed access publication – behind paywall"></a></span></span></span>
</li>
<li id="cite_note-Buhmann-8"><span class="mw-cite-backlink"><b><a href="#cite_ref-Buhmann_8-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFBuhmannKuhnel1992" class="citation book cs1">Buhmann, J.; Kuhnel, H. (1992). "Unsupervised and supervised data clustering with competitive neural networks". <i>[Proceedings 1992] IJCNN International Joint Conference on Neural Networks</i>. Vol. 4. IEEE. pp. <span class="nowrap">796–</span>801. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1109%2Fijcnn.1992.227220">10.1109/ijcnn.1992.227220</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>0780305590</bdi>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a> <a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:62651220">62651220</a>.</cite></span>
</li>
<li id="cite_note-Comesana-9"><span class="mw-cite-backlink"><b><a href="#cite_ref-Comesana_9-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFComesaña-CamposBouza-Rodríguez2016" class="citation journal cs1">Comesaña-Campos, Alberto; Bouza-Rodríguez, José Benito (June 2016). "An application of Hebbian learning in the design process decision-making". <i>Journal of Intelligent Manufacturing</i>. <b>27</b> (3): <span class="nowrap">487–</span>506. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1007%2Fs10845-014-0881-z">10.1007/s10845-014-0881-z</a>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a> <a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/0956-5515">0956-5515</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a> <a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:207171436">207171436</a>.</cite></span>
</li>
<li id="cite_note-Carpenter-10"><span class="mw-cite-backlink"><b><a href="#cite_ref-Carpenter_10-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFCarpenter,_G.A.Grossberg,_S.1988" class="citation journal cs1">Carpenter, G.A. & Grossberg, S. (1988). <a rel="nofollow" class="external text" href="https://web.archive.org/web/20180516131553/http://www.cns.bu.edu/Profiles/Grossberg/CarGro1988Computer.pdf">"The ART of adaptive pattern recognition by a self-organizing neural network"</a> <span class="cs1-format">(PDF)</span>. <i>Computer</i>. <b>21</b> (3): <span class="nowrap">77–</span>88. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1109%2F2.33">10.1109/2.33</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a> <a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:14625094">14625094</a>. Archived from <a rel="nofollow" class="external text" href="http://www.cns.bu.edu/Profiles/Grossberg/CarGro1988Computer.pdf">the original</a> <span class="cs1-format">(PDF)</span> on 2018-05-16<span class="reference-accessdate">. Retrieved <span class="nowrap">2013-09-16</span></span>.</cite></span>
</li>
<li id="cite_note-tds-ul-11"><span class="mw-cite-backlink"><b><a href="#cite_ref-tds-ul_11-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFRoman2019" class="citation web cs1">Roman, Victor (2019-04-21). <a rel="nofollow" class="external text" href="https://towardsdatascience.com/unsupervised-machine-learning-clustering-analysis-d40f2b34ae7e">"Unsupervised Machine Learning: Clustering Analysis"</a>. <i>Medium</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20200821132257/https://towardsdatascience.com/unsupervised-machine-learning-clustering-analysis-d40f2b34ae7e">Archived</a> from the original on 2020-08-21<span class="reference-accessdate">. Retrieved <span class="nowrap">2019-10-01</span></span>.</cite></span>
</li>
<li id="cite_note-JordanBishop2004-12"><span class="mw-cite-backlink"><b><a href="#cite_ref-JordanBishop2004_12-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFJordanBishop2004" class="citation book cs1">Jordan, Michael I.; Bishop, Christopher M. (2004). "7. Intelligent Systems §Neural Networks". In Tucker, Allen B. (ed.). <a rel="nofollow" class="external text" href="https://www.taylorfrancis.com/books/mono/10.1201/9780203494455/computer-science-handbook-allen-tucker"><i>Computer Science Handbook</i></a> (2nd ed.). Chapman & Hall/CRC Press. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1201%2F9780203494455">10.1201/9780203494455</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>1-58488-360-X</bdi>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20221103234201/https://www.taylorfrancis.com/books/mono/10.1201/9780203494455/computer-science-handbook-allen-tucker">Archived</a> from the original on 2022-11-03<span class="reference-accessdate">. Retrieved <span class="nowrap">2022-11-03</span></span>.</cite></span>
</li>
<li id="cite_note-Hastie-13"><span class="mw-cite-backlink"><b><a href="#cite_ref-Hastie_13-0">^</a></b></span> <span class="reference-text"><a href="#CITEREFHastieTibshiraniFriedman2009">Hastie, Tibshirani & Friedman 2009</a>, pp. 485–586</span>
</li>
<li id="cite_note-tds-kmeans-14"><span class="mw-cite-backlink"><b><a href="#cite_ref-tds-kmeans_14-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFGarbade2018" class="citation web cs1">Garbade, Dr Michael J. (2018-09-12). <a rel="nofollow" class="external text" href="https://towardsdatascience.com/understanding-k-means-clustering-in-machine-learning-6a6e67336aa1">"Understanding K-means Clustering in Machine Learning"</a>. <i>Medium</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20190528183913/https://towardsdatascience.com/understanding-k-means-clustering-in-machine-learning-6a6e67336aa1">Archived</a> from the original on 2019-05-28<span class="reference-accessdate">. Retrieved <span class="nowrap">2019-10-31</span></span>.</cite></span>
</li>
<li id="cite_note-TensorLVMs-15"><span class="mw-cite-backlink"><b><a href="#cite_ref-TensorLVMs_15-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFAnandkumarGeHsuKakade2014" class="citation journal cs1">Anandkumar, Animashree; Ge, Rong; Hsu, Daniel; Kakade, Sham; Telgarsky, Matus (2014). <a rel="nofollow" class="external text" href="http://www.jmlr.org/papers/volume15/anandkumar14b/anandkumar14b.pdf">"Tensor Decompositions for Learning Latent Variable Models"</a> <span class="cs1-format">(PDF)</span>. <i>Journal of Machine Learning Research</i>. <b>15</b>: <span class="nowrap">2773–</span>2832. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1210.7559">1210.7559</a></span>. <a href="Bibcode_(identifier)" class="mw-redirect" title="Bibcode (identifier)">Bibcode</a>:<a rel="nofollow" class="external text" href="https://ui.adsabs.harvard.edu/abs/2012arXiv1210.7559A">2012arXiv1210.7559A</a>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20150320201108/http://jmlr.org/papers/volume15/anandkumar14b/anandkumar14b.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2015-03-20<span class="reference-accessdate">. Retrieved <span class="nowrap">2015-04-10</span></span>.</cite></span>
</li>
</ol></div></div>
<div class="mw-heading mw-heading2"><h2 id="Further_reading">Further reading</h2></div>
<style data-mw-deduplicate="TemplateStyles:r1239549316">
/* start https://en.wikipedia.org/ */
.mw-parser-output .refbegin{margin-bottom:0.5em}.mw-parser-output .refbegin-hanging-indents>ul{margin-left:0}.mw-parser-output .refbegin-hanging-indents>ul>li{margin-left:0;padding-left:3.2em;text-indent:-3.2em}.mw-parser-output .refbegin-hanging-indents ul,.mw-parser-output .refbegin-hanging-indents ul li{list-style:none}@media(max-width:720px){.mw-parser-output .refbegin-hanging-indents>ul>li{padding-left:1.6em;text-indent:-1.6em}}.mw-parser-output .refbegin-columns{margin-top:0.3em}.mw-parser-output .refbegin-columns ul{margin-top:0}.mw-parser-output .refbegin-columns li{page-break-inside:avoid;break-inside:avoid-column}@media screen{.mw-parser-output .refbegin{font-size:90%}}
/* end https://en.wikipedia.org/ */
</style><div class="refbegin" style="">
<ul><li><cite id="CITEREFBousquet,_O.von_Luxburg,_U.Raetsch,_G.2004" class="citation book cs1">Bousquet, O.; <a href="Ulrike_von_Luxburg" title="Ulrike von Luxburg">von Luxburg, U.</a>; Raetsch, G., eds. (2004). <a rel="nofollow" class="external text" href="https://archive.org/details/springer_10.1007-b100712"><i>Advanced Lectures on Machine Learning</i></a>. Springer. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>978-3540231226</bdi>.</cite></li>
<li><cite id="CITEREFDuda,_Richard_O.Hart,_Peter_E.Stork,_David_G.2001" class="citation book cs1"><a href="Richard_O._Duda" title="Richard O. Duda">Duda, Richard O.</a>; <a href="Peter_E._Hart" title="Peter E. Hart">Hart, Peter E.</a>; Stork, David G. (2001). "Unsupervised Learning and Clustering". <a href="Pattern_classification" class="mw-redirect" title="Pattern classification"><i>Pattern classification</i></a> (2nd ed.). Wiley. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>0-471-05669-3</bdi>.</cite></li>
<li><cite id="CITEREFHastieTibshiraniFriedman2009" class="citation book cs1"><a href="Trevor_Hastie" title="Trevor Hastie">Hastie, Trevor</a>; <a href="Robert_Tibshirani" title="Robert Tibshirani">Tibshirani, Robert</a>; Friedman, Jerome (2009). <a rel="nofollow" class="external text" href="https://link.springer.com/chapter/10.1007/978-0-387-84858-7_14">"Unsupervised Learning"</a>. <i>The Elements of Statistical Learning: Data mining, Inference, and Prediction</i>. Springer. pp. <span class="nowrap">485–</span>586. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<a rel="nofollow" class="external text" href="https://doi.org/10.1007%2F978-0-387-84858-7_14">10.1007/978-0-387-84858-7_14</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>978-0-387-84857-0</bdi>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20221103234204/https://link.springer.com/chapter/10.1007/978-0-387-84858-7_14">Archived</a> from the original on 2022-11-03<span class="reference-accessdate">. Retrieved <span class="nowrap">2022-11-03</span></span>.</cite></li>
<li><cite id="CITEREFHintonSejnowski1999" class="citation book cs1"><a href="Geoffrey_Hinton" title="Geoffrey Hinton">Hinton, Geoffrey</a>; <a href="Terrence_J._Sejnowski" class="mw-redirect" title="Terrence J. Sejnowski">Sejnowski, Terrence J.</a>, eds. (1999). <i>Unsupervised Learning: Foundations of Neural Computation</i>. <a href="MIT_Press" title="MIT Press">MIT Press</a>. <a href="ISBN_(identifier)" class="mw-redirect" title="ISBN (identifier)">ISBN</a> <bdi>0-262-58168-X</bdi>.</cite></li></ul>
</div>
<div class="navbox-styles"><style data-mw-deduplicate="TemplateStyles:r1236075235">
/* start https://en.wikipedia.org/ */
.mw-parser-output .navbox{box-sizing:border-box;border:1px solid #a2a9b1;width:100%;clear:both;font-size:88%;text-align:center;padding:1px;margin:1em auto 0}.mw-parser-output .navbox .navbox{margin-top:0}.mw-parser-output .navbox+.navbox,.mw-parser-output .navbox+.navbox-styles+.navbox{margin-top:-1px}.mw-parser-output .navbox-inner,.mw-parser-output .navbox-subgroup{width:100%}.mw-parser-output .navbox-group,.mw-parser-output .navbox-title,.mw-parser-output .navbox-abovebelow{padding:0.25em 1em;line-height:1.5em;text-align:center}.mw-parser-output .navbox-group{white-space:nowrap;text-align:right}.mw-parser-output .navbox,.mw-parser-output .navbox-subgroup{background-color:#fdfdfd}.mw-parser-output .navbox-list{line-height:1.5em;border-color:#fdfdfd}.mw-parser-output .navbox-list-with-group{text-align:left;border-left-width:2px;border-left-style:solid}.mw-parser-output tr+tr>.navbox-abovebelow,.mw-parser-output tr+tr>.navbox-group,.mw-parser-output tr+tr>.navbox-image,.mw-parser-output tr+tr>.navbox-list{border-top:2px solid #fdfdfd}.mw-parser-output .navbox-title{background-color:#ccf}.mw-parser-output .navbox-abovebelow,.mw-parser-output .navbox-group,.mw-parser-output .navbox-subgroup .navbox-title{background-color:#ddf}.mw-parser-output .navbox-subgroup .navbox-group,.mw-parser-output .navbox-subgroup .navbox-abovebelow{background-color:#e6e6ff}.mw-parser-output .navbox-even{background-color:#f7f7f7}.mw-parser-output .navbox-odd{background-color:transparent}.mw-parser-output .navbox .hlist td dl,.mw-parser-output .navbox .hlist td ol,.mw-parser-output .navbox .hlist td ul,.mw-parser-output .navbox td.hlist dl,.mw-parser-output .navbox td.hlist ol,.mw-parser-output .navbox td.hlist ul{padding:0.125em 0}.mw-parser-output .navbox .navbar{display:block;font-size:100%}.mw-parser-output .navbox-title .navbar{float:left;text-align:left;margin-right:0.5em}body.skin--responsive .mw-parser-output .navbox-image img{max-width:none!important}@media print{body.ns-0 .mw-parser-output .navbox{display:none!important}}
/* end https://en.wikipedia.org/ */
</style></div><div role="navigation" class="navbox" aria-labelledby="Differentiable_computing254" style="padding:3px"><table class="nowraplinks hlist mw-collapsible autocollapse navbox-inner" style="border-spacing:0;background:transparent;color:inherit"><tbody><tr><th scope="col" class="navbox-title" colspan="2"><div id="Differentiable_computing254" style="font-size:114%;margin:0 4em">Differentiable computing</div></th></tr><tr><th scope="row" class="navbox-group" style="width:1%"><a href="Differentiable_function" title="Differentiable function">General</a></th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><b><a href="Differentiable_programming" title="Differentiable programming">Differentiable programming</a></b></li>
<li><a href="Information_geometry" title="Information geometry">Information geometry</a></li>
<li><a href="Statistical_manifold" title="Statistical manifold">Statistical manifold</a></li>
<li><a href="Automatic_differentiation" title="Automatic differentiation">Automatic differentiation</a></li>
<li><a href="Neuromorphic_computing" title="Neuromorphic computing">Neuromorphic computing</a></li>
<li><a href="Pattern_recognition" title="Pattern recognition">Pattern recognition</a></li>
<li><a href="Ricci_calculus" title="Ricci calculus">Ricci calculus</a></li>
<li><a href="Computational_learning_theory" title="Computational learning theory">Computational learning theory</a></li>
<li><a href="Inductive_bias" title="Inductive bias">Inductive bias</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Hardware</th><td class="navbox-list-with-group navbox-list navbox-even" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="Graphcore" title="Graphcore">IPU</a></li>
<li><a href="Tensor_Processing_Unit" title="Tensor Processing Unit">TPU</a></li>
<li><a href="Vision_processing_unit" title="Vision processing unit">VPU</a></li>
<li><a href="Memristor" title="Memristor">Memristor</a></li>
<li><a href="SpiNNaker" title="SpiNNaker">SpiNNaker</a></li></ul>
</div></td></tr><tr><th scope="row" class="navbox-group" style="width:1%">Software libraries</th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em">
<ul><li><a href="TensorFlow" title="TensorFlow">TensorFlow</a></li>
<li><a href="PyTorch" title="PyTorch">PyTorch</a></li>
<li><a href="Keras" title="Keras">Keras</a></li>
<li><a href="Scikit-learn" title="Scikit-learn">scikit-learn</a></li>
<li><a href="Theano_(software)" title="Theano (software)">Theano</a></li>
<li><a href="JAX_(software)" title="JAX (software)">JAX</a></li>
<li><a href="Flux_(machine-learning_framework)" title="Flux (machine-learning framework)">Flux.jl</a></li>
<li><a href="MindSpore" title="MindSpore">MindSpore</a></li></ul>
</div></td></tr><tr><td class="navbox-abovebelow" colspan="2"><div>
<ul><li><span class="noviewer" typeof="mw:File"></span> Portals
<ul><li><a href="Portal%3AComputer_programming" title="Portal:Computer programming">Computer programming</a></li>
<li><a href="Portal%3ATechnology" title="Portal:Technology">Technology</a></li></ul></li></ul>
</div></td></tr></tbody></table></div>
<div class="navbox-styles"></div><div role="navigation" class="navbox authority-control" aria-label="Navbox391" style="padding:3px"><table class="nowraplinks hlist navbox-inner" style="border-spacing:0;background:transparent;color:inherit"><tbody><tr><th scope="row" class="navbox-group" style="width:1%">Authority control databases: National </th><td class="navbox-list-with-group navbox-list navbox-odd" style="width:100%;padding:0"><div style="padding:0 0.25em"><ul><li><span class="uid"><a rel="nofollow" class="external text" href="https://d-nb.info/gnd/4580265-8">Germany</a></span></li></ul></div></td></tr></tbody></table></div></div><!--htdig_noindex--><div><div class="zim-footer">
This article is issued from <a class="external text" title="Last edited on 2025-07-16" href="https://en.wikipedia.org/wiki/?title=Unsupervised_learning&oldid=1300836256">Wikipedia</a>. The text is available under <a class="external text" href="https://creativecommons.org/licenses/by-sa/4.0/deed.en">Creative Commons Attribution-Share Alike 4.0</a> unless otherwise noted. Additional terms may apply for the media files.
</div>
</div><!--/htdig_noindex--></div>
</div>
</main>
</div>
</div>
</div>
</body></html>